agora inbox for pgsql-hackers@postgresql.orghelp / color / mirror / Atom feed
[PATCH v1 4/4] Extra tests 386+ messages / 2 participants [nested] [flat]
* [PATCH v1 4/4] Extra tests @ 2025-08-31 09:10 Nikolay Shaplov <dhyan@nataraj.su> 0 siblings, 0 replies; 386+ messages in thread From: Nikolay Shaplov @ 2025-08-31 09:10 UTC (permalink / raw) Add more tests for trenary reloptions in dummy_index_am module --- src/test/modules/dummy_index_am/README | 1 + .../modules/dummy_index_am/dummy_index_am.c | 36 +++++++++++++----- .../dummy_index_am/expected/reloptions.out | 38 +++++++++++++++++-- .../modules/dummy_index_am/sql/reloptions.sql | 16 ++++++++ src/test/regress/expected/reloptions.out | 4 +- src/test/regress/sql/reloptions.sql | 4 +- 6 files changed, 81 insertions(+), 18 deletions(-) diff --git a/src/test/modules/dummy_index_am/README b/src/test/modules/dummy_index_am/README index 61510f02fae..78d1d84ed9a 100644 --- a/src/test/modules/dummy_index_am/README +++ b/src/test/modules/dummy_index_am/README @@ -6,6 +6,7 @@ access method, whose code is kept a maximum simple. This includes tests for all relation option types: - boolean +- trenary - enum - integer - real diff --git a/src/test/modules/dummy_index_am/dummy_index_am.c b/src/test/modules/dummy_index_am/dummy_index_am.c index 94ef639b6fc..19826e50e6d 100644 --- a/src/test/modules/dummy_index_am/dummy_index_am.c +++ b/src/test/modules/dummy_index_am/dummy_index_am.c @@ -22,7 +22,7 @@ PG_MODULE_MAGIC; /* parse table for fillRelOptions */ -static relopt_parse_elt di_relopt_tab[6]; +static relopt_parse_elt di_relopt_tab[8]; /* Kind of relation options for dummy index */ static relopt_kind di_relopt_kind; @@ -40,6 +40,8 @@ typedef struct DummyIndexOptions int option_int; double option_real; bool option_bool; + trenary option_trenary1; + trenary option_trenary2; DummyAmEnum option_enum; int option_string_val_offset; int option_string_null_offset; @@ -96,23 +98,37 @@ create_reloptions_table(void) di_relopt_tab[2].opttype = RELOPT_TYPE_BOOL; di_relopt_tab[2].offset = offsetof(DummyIndexOptions, option_bool); + add_trenary_reloption(di_relopt_kind, "option_trenary1", + "First trenary option for dummy_index_am", + TRENARY_UNSET, NULL, AccessExclusiveLock); + di_relopt_tab[3].optname = "option_trenary1"; + di_relopt_tab[3].opttype = RELOPT_TYPE_TRENARY; + di_relopt_tab[3].offset = offsetof(DummyIndexOptions, option_trenary1); + + add_trenary_reloption(di_relopt_kind, "option_trenary2", + "Second trenary option for dummy_index_am", + TRENARY_TRUE, "do_not_know_yet", AccessExclusiveLock); + di_relopt_tab[4].optname = "option_trenary2"; + di_relopt_tab[4].opttype = RELOPT_TYPE_TRENARY; + di_relopt_tab[4].offset = offsetof(DummyIndexOptions, option_trenary2); + add_enum_reloption(di_relopt_kind, "option_enum", "Enum option for dummy_index_am", dummyAmEnumValues, DUMMY_AM_ENUM_ONE, "Valid values are \"one\" and \"two\".", AccessExclusiveLock); - di_relopt_tab[3].optname = "option_enum"; - di_relopt_tab[3].opttype = RELOPT_TYPE_ENUM; - di_relopt_tab[3].offset = offsetof(DummyIndexOptions, option_enum); + di_relopt_tab[5].optname = "option_enum"; + di_relopt_tab[5].opttype = RELOPT_TYPE_ENUM; + di_relopt_tab[5].offset = offsetof(DummyIndexOptions, option_enum); add_string_reloption(di_relopt_kind, "option_string_val", "String option for dummy_index_am with non-NULL default", "DefaultValue", &validate_string_option, AccessExclusiveLock); - di_relopt_tab[4].optname = "option_string_val"; - di_relopt_tab[4].opttype = RELOPT_TYPE_STRING; - di_relopt_tab[4].offset = offsetof(DummyIndexOptions, + di_relopt_tab[6].optname = "option_string_val"; + di_relopt_tab[6].opttype = RELOPT_TYPE_STRING; + di_relopt_tab[6].offset = offsetof(DummyIndexOptions, option_string_val_offset); /* @@ -123,9 +139,9 @@ create_reloptions_table(void) NULL, /* description */ NULL, &validate_string_option, AccessExclusiveLock); - di_relopt_tab[5].optname = "option_string_null"; - di_relopt_tab[5].opttype = RELOPT_TYPE_STRING; - di_relopt_tab[5].offset = offsetof(DummyIndexOptions, + di_relopt_tab[7].optname = "option_string_null"; + di_relopt_tab[7].opttype = RELOPT_TYPE_STRING; + di_relopt_tab[7].offset = offsetof(DummyIndexOptions, option_string_null_offset); } diff --git a/src/test/modules/dummy_index_am/expected/reloptions.out b/src/test/modules/dummy_index_am/expected/reloptions.out index c873a80bb75..bcf164f5bf7 100644 --- a/src/test/modules/dummy_index_am/expected/reloptions.out +++ b/src/test/modules/dummy_index_am/expected/reloptions.out @@ -18,6 +18,8 @@ SET client_min_messages TO 'notice'; CREATE INDEX dummy_test_idx ON dummy_test_tab USING dummy_index_am (i) WITH ( option_bool = false, + option_trenary1, + option_trenary2 = off, option_int = 5, option_real = 3.1, option_enum = 'two', @@ -31,16 +33,20 @@ SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; unnest ------------------------ option_bool=false + option_trenary1=true + option_trenary2=off option_int=5 option_real=3.1 option_enum=two option_string_val=null option_string_null=val -(6 rows) +(8 rows) -- ALTER INDEX .. SET ALTER INDEX dummy_test_idx SET (option_int = 10); ALTER INDEX dummy_test_idx SET (option_bool = true); +ALTER INDEX dummy_test_idx SET (option_trenary1 = false); +ALTER INDEX dummy_test_idx SET (option_trenary2 = Do_Not_Know_YET); ALTER INDEX dummy_test_idx SET (option_real = 3.2); ALTER INDEX dummy_test_idx SET (option_string_val = 'val2'); ALTER INDEX dummy_test_idx SET (option_string_null = NULL); @@ -49,19 +55,23 @@ ALTER INDEX dummy_test_idx SET (option_enum = 'three'); ERROR: invalid value for enum option "option_enum": three DETAIL: Valid values are "one" and "two". SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; - unnest -------------------------- + unnest +--------------------------------- option_int=10 option_bool=true + option_trenary1=false + option_trenary2=do_not_know_yet option_real=3.2 option_string_val=val2 option_string_null=null option_enum=one -(6 rows) +(8 rows) -- ALTER INDEX .. RESET ALTER INDEX dummy_test_idx RESET (option_int); ALTER INDEX dummy_test_idx RESET (option_bool); +ALTER INDEX dummy_test_idx RESET (option_trenary1); +ALTER INDEX dummy_test_idx RESET (option_trenary2); ALTER INDEX dummy_test_idx RESET (option_real); ALTER INDEX dummy_test_idx RESET (option_enum); ALTER INDEX dummy_test_idx RESET (option_string_val); @@ -100,6 +110,26 @@ SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; (1 row) ALTER INDEX dummy_test_idx RESET (option_bool); +-- Trenary +ALTER INDEX dummy_test_idx SET (option_trenary1 = 4); -- error +ERROR: invalid value for trenary option "option_trenary1": 4 +ALTER INDEX dummy_test_idx SET (option_trenary1 = 1); -- ok, as true +ALTER INDEX dummy_test_idx SET (option_trenary1 = 3.4); -- error +ERROR: invalid value for trenary option "option_trenary1": 3.4 +ALTER INDEX dummy_test_idx SET (option_trenary1 = 'val4'); -- error +ERROR: invalid value for trenary option "option_trenary1": val4 +ALTER INDEX dummy_test_idx SET (option_trenary1 = 'do_not_know_yet'); -- error. Valid for trenary2 not for trenary1 +ERROR: invalid value for trenary option "option_trenary1": do_not_know_yet +ALTER INDEX dummy_test_idx SET (option_trenary2 = 'do_not_know_yet'); -- ok +SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; + unnest +--------------------------------- + option_trenary1=1 + option_trenary2=do_not_know_yet +(2 rows) + +ALTER INDEX dummy_test_idx RESET (option_trenary1); +ALTER INDEX dummy_test_idx RESET (option_trenary2); -- Float ALTER INDEX dummy_test_idx SET (option_real = 4); -- ok ALTER INDEX dummy_test_idx SET (option_real = true); -- error diff --git a/src/test/modules/dummy_index_am/sql/reloptions.sql b/src/test/modules/dummy_index_am/sql/reloptions.sql index 6749d763e6a..7e752b12d16 100644 --- a/src/test/modules/dummy_index_am/sql/reloptions.sql +++ b/src/test/modules/dummy_index_am/sql/reloptions.sql @@ -18,6 +18,8 @@ SET client_min_messages TO 'notice'; CREATE INDEX dummy_test_idx ON dummy_test_tab USING dummy_index_am (i) WITH ( option_bool = false, + option_trenary1, + option_trenary2 = off, option_int = 5, option_real = 3.1, option_enum = 'two', @@ -30,6 +32,8 @@ SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; -- ALTER INDEX .. SET ALTER INDEX dummy_test_idx SET (option_int = 10); ALTER INDEX dummy_test_idx SET (option_bool = true); +ALTER INDEX dummy_test_idx SET (option_trenary1 = false); +ALTER INDEX dummy_test_idx SET (option_trenary2 = Do_Not_Know_YET); ALTER INDEX dummy_test_idx SET (option_real = 3.2); ALTER INDEX dummy_test_idx SET (option_string_val = 'val2'); ALTER INDEX dummy_test_idx SET (option_string_null = NULL); @@ -40,6 +44,8 @@ SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; -- ALTER INDEX .. RESET ALTER INDEX dummy_test_idx RESET (option_int); ALTER INDEX dummy_test_idx RESET (option_bool); +ALTER INDEX dummy_test_idx RESET (option_trenary1); +ALTER INDEX dummy_test_idx RESET (option_trenary2); ALTER INDEX dummy_test_idx RESET (option_real); ALTER INDEX dummy_test_idx RESET (option_enum); ALTER INDEX dummy_test_idx RESET (option_string_val); @@ -60,6 +66,16 @@ ALTER INDEX dummy_test_idx SET (option_bool = 3.4); -- error ALTER INDEX dummy_test_idx SET (option_bool = 'val4'); -- error SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; ALTER INDEX dummy_test_idx RESET (option_bool); +-- Trenary +ALTER INDEX dummy_test_idx SET (option_trenary1 = 4); -- error +ALTER INDEX dummy_test_idx SET (option_trenary1 = 1); -- ok, as true +ALTER INDEX dummy_test_idx SET (option_trenary1 = 3.4); -- error +ALTER INDEX dummy_test_idx SET (option_trenary1 = 'val4'); -- error +ALTER INDEX dummy_test_idx SET (option_trenary1 = 'do_not_know_yet'); -- error. Valid for trenary2 not for trenary1 +ALTER INDEX dummy_test_idx SET (option_trenary2 = 'do_not_know_yet'); -- ok +SELECT unnest(reloptions) FROM pg_class WHERE relname = 'dummy_test_idx'; +ALTER INDEX dummy_test_idx RESET (option_trenary1); +ALTER INDEX dummy_test_idx RESET (option_trenary2); -- Float ALTER INDEX dummy_test_idx SET (option_real = 4); -- ok ALTER INDEX dummy_test_idx SET (option_real = true); -- error diff --git a/src/test/regress/expected/reloptions.out b/src/test/regress/expected/reloptions.out index e32431ca067..54b3424ef2b 100644 --- a/src/test/regress/expected/reloptions.out +++ b/src/test/regress/expected/reloptions.out @@ -98,7 +98,7 @@ SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; {fillfactor=13,autovacuum_enabled=false} (1 row) --- Tests for future (FIXME) trenary options +-- Tests for trenary options -- behave as boolean option: accept unassigned name and truncated value DROP TABLE reloptions_test; CREATE TABLE reloptions_test(i INT) WITH (vacuum_truncate); @@ -116,7 +116,7 @@ SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; {vacuum_truncate=fals} (1 row) --- preferred "true" alias is used when storing +-- preferred "true" alias is stored in pg_class DROP TABLE reloptions_test; CREATE TABLE reloptions_test(i INT) WITH (vacuum_index_cleanup=on); SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; diff --git a/src/test/regress/sql/reloptions.sql b/src/test/regress/sql/reloptions.sql index 051142eed5f..1971d460672 100644 --- a/src/test/regress/sql/reloptions.sql +++ b/src/test/regress/sql/reloptions.sql @@ -59,7 +59,7 @@ UPDATE pg_class ALTER TABLE reloptions_test RESET (illegal_option); SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; --- Tests for future (FIXME) trenary options +-- Tests for trenary options -- behave as boolean option: accept unassigned name and truncated value DROP TABLE reloptions_test; @@ -70,7 +70,7 @@ DROP TABLE reloptions_test; CREATE TABLE reloptions_test(i INT) WITH (vacuum_truncate=FaLS); SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; --- preferred "true" alias is used when storing +-- preferred "true" alias is stored in pg_class DROP TABLE reloptions_test; CREATE TABLE reloptions_test(i INT) WITH (vacuum_index_cleanup=on); SELECT reloptions FROM pg_class WHERE oid = 'reloptions_test'::regclass; -- 2.39.2 --nextPart6784952.tM3a2QDmDi-- --nextPart6187903.UjTJXf6HLC Content-Type: application/pgp-signature; name="signature.asc" Content-Description: This is a digitally signed message part. Content-Transfer-Encoding: 7Bit -----BEGIN PGP SIGNATURE----- iQEzBAABCgAdFiEE+sk3ebqQKlezKOi8PMbfuIHAGpgFAmi0gCMACgkQPMbfuIHA Gpi3Dgf/dQxIJ7WM8rL4szT+d+7gCw1BUzA7H4SkLtGM8E+akUvbjt6lFpCKN7fP dHWTY7NeHsLy8hXB7Z+P4edaHAqNQ7pXtESuC0jBB1Ta5r9f0mUz/tNvyaoj1YPV SES411gQwABIJ/ZXoqrcAHgmJoyAKZcFWU9HBAhJyL3Da/o7L86wgro4a6ov7cgP 1mJa6WjH1qf1v4PPrZYxLMmU5v3uacNDFowNsQ8+uHs+slUWjkw4VYatdReWla0a 4qBnMaht1Xky3f55YfB7Gx28CBUUPMREUVKFOrcopq95kZ9WASoTJfU9Y80IDPWb KVcJvYDk5TEL1t48GpFkjZbL9od51w== =yrfm -----END PGP SIGNATURE----- --nextPart6187903.UjTJXf6HLC-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
* [PATCH] Teach REPACK to upgrade its lock safely. @ 2026-04-14 09:59 Antonin Houska <ah@cybertec.at> 0 siblings, 0 replies; 386+ messages in thread From: Antonin Houska @ 2026-04-14 09:59 UTC (permalink / raw) REPACK (CONCURRENTLY) needs to upgrade its ShareUpdateExclusiveLock (SUEL) to AccessExclusiveLock at the end of table processing. If another session, which already has a non-conflicting lock on the table, tried to get SUEL too, it would end up in a deadlock with REPACK. Such situation does not hurt as long as the deadlock detector choses to terminate the other session, but it's possible that it terminates REPACK. A lot of REPACK's work would be wasted this way. This patch checks for such situation, and if the other session tries to get the SUEL, it receives a deadlock report before it actually starts waiting. Please note that the other session can safely run VACUUM w/o receiving the deadlock report. The point is that VACUUM cannot run in a transaction block, so the session cannot have any other lock when trying to get SUEL on the table. In that case, REPACK gets its AccessExclusiveLock first, even if VACUUM requested SUEL earlier. VACUUM then just needs wait for REPACK to finish, and then it can start. --- src/backend/commands/repack.c | 25 +++++++- src/backend/storage/lmgr/deadlock.c | 10 ++-- src/backend/storage/lmgr/lock.c | 30 ++++++++++ src/backend/storage/lmgr/proc.c | 55 ++++++++++++++++- src/include/storage/lock.h | 5 +- src/include/storage/proc.h | 6 +- .../injection_points/expected/repack.out | 59 ++++++++++++++++++- .../injection_points/specs/repack.spec | 29 +++++++++ 8 files changed, 209 insertions(+), 10 deletions(-) diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c index 58e3867246f..e9193067666 100644 --- a/src/backend/commands/repack.c +++ b/src/backend/commands/repack.c @@ -285,6 +285,18 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) * to understand and we don't lose any functionality. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); + + /* + * Also set the PROC_IN_CONCURRENT_REPACK flag. This makes the lock + * manager cause anyone that would conflict with us to error out. + * It's important to set this flag ahead of actually locking the + * relation; it won't of course affect anyone until we do have a lock + * that others can conflict with. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags |= PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); } /* @@ -3086,7 +3098,18 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap, LockRelationOid(OldHeap->rd_rel->reltoastrelid, AccessExclusiveLock); /* - * Tuples and pages of the old heap will be gone, but the heap will stay. + * Now that we have all access-exclusive locks on all relations, we no + * longer want other processes to error out when trying to acquire a + * conflicting lock. Therefore, unset our flag. + */ + LWLockAcquire(ProcArrayLock, LW_EXCLUSIVE); + MyProc->statusFlags &= ~PROC_IN_CONCURRENT_REPACK; + ProcGlobal->statusFlags[MyProc->pgxactoff] = MyProc->statusFlags; + LWLockRelease(ProcArrayLock); + + /* + * Tuples and pages of the old heap will be gone, but the heap itself will + * stay. */ TransferPredicateLocksToHeapRelation(OldHeap); foreach_ptr(RelationData, index, indexrels) diff --git a/src/backend/storage/lmgr/deadlock.c b/src/backend/storage/lmgr/deadlock.c index b8962d875b6..3772a7aa3df 100644 --- a/src/backend/storage/lmgr/deadlock.c +++ b/src/backend/storage/lmgr/deadlock.c @@ -1146,8 +1146,10 @@ DeadLockReport(void) void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, - LOCK *lock, - PGPROC *proc2) + LOCK *lock, /* XXX Only lockmode? */ + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2) { DEADLOCK_INFO *info = &deadlockDetails[0]; @@ -1155,8 +1157,8 @@ RememberSimpleDeadLock(PGPROC *proc1, info->lockmode = lockmode; info->pid = proc1->pid; info++; - info->locktag = proc2->waitLock->tag; - info->lockmode = proc2->waitLockMode; + info->locktag = *locktag2; + info->lockmode = lockmode2; info->pid = proc2->pid; nDeadlockDetails = 2; } diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c index c221fe96889..44aa2ca26a7 100644 --- a/src/backend/storage/lmgr/lock.c +++ b/src/backend/storage/lmgr/lock.c @@ -674,6 +674,36 @@ LockHeldByMe(const LOCKTAG *locktag, return false; } +/* + * HaveRelLock -- test whether the current backend has a fast path lock on + * the relation in any mode. + */ +bool +HaveFastPathLock(Oid relid) +{ + uint32 group = FAST_PATH_REL_GROUP(relid); + bool result = false; + + LWLockAcquire(&MyProc->fpInfoLock, LW_SHARED); + for (int i = 0; i < FP_LOCK_SLOTS_PER_GROUP; i++) + { + uint32 f = FAST_PATH_SLOT(group, i); + uint32 lockmask; + + if (relid != MyProc->fpRelId[f]) + continue; + lockmask = FAST_PATH_GET_BITS(MyProc, f); + if (lockmask) + { + result = true; + break; + } + } + LWLockRelease(&MyProc->fpInfoLock); + + return result; +} + #ifdef USE_ASSERT_CHECKING /* * GetLockMethodLocalHash -- return the hash of local locks, for modules that diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c index 1ac25068d62..3d47b1284a3 100644 --- a/src/backend/storage/lmgr/proc.c +++ b/src/backend/storage/lmgr/proc.c @@ -1156,6 +1156,7 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) LOCKMASK myHeldLocks; bool early_deadlock = false; PGPROC *leader = MyProc->lockGroupLeader; + bool check_repack = false; Assert(LWLockHeldByMeInMode(partitionLock, LW_EXCLUSIVE)); @@ -1193,6 +1194,56 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) } } + /* + * If I am already holding this lock (in any mode) and trying to get + * ShareUpdateExclusiveLock (or higher), check if REPACK (CONCURRENTLY) is + * already holding ShareUpdateExclusiveLock. The problem is if that by + * trying to upgrade the lock to AccessExclusiveLock it would get into a + * deadlock with us. + */ + if (lock->tag.locktag_type == LOCKTAG_RELATION && + lock->tag.locktag_field1 == MyDatabaseId && + lockmode >= ShareUpdateExclusiveLock) + { + if (myHeldLocks > 0 || + HaveFastPathLock(lock->tag.locktag_field2)) + check_repack = true; + } + if (check_repack) + { + dlist_iter iter; + + dlist_foreach(iter, &lock->procLocks) + { + PROCLOCK *otherproclock; + PGPROC *otherproc; + + otherproclock = dlist_container(PROCLOCK, lockLink, iter.cur); + otherproc = otherproclock->tag.myProc; + if (otherproc != MyProc && + otherproc->statusFlags & PROC_IN_CONCURRENT_REPACK) + { + LOCKMASK repackmask = otherproclock->holdMask; + + /* + * Is REPACK already holding ShareUpdateExclusiveLock? + */ + if ((repackmask & LOCKBIT_ON(ShareUpdateExclusiveLock)) != 0) + { + /* + * There should be no more than one REPACK working on + * particular table, so let's error out. + */ + RememberSimpleDeadLock(MyProc, lockmode, lock, + otherproc, + ShareUpdateExclusiveLock, + &otherproclock->tag.myLock->tag); + return PROC_WAIT_STATUS_ERROR; + } + } + } + } + /* * Determine where to add myself in the wait queue. * @@ -1240,7 +1291,9 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait) * a flag to check below, and break out of loop. Also, * record deadlock info for later message. */ - RememberSimpleDeadLock(MyProc, lockmode, lock, proc); + RememberSimpleDeadLock(MyProc, lockmode, lock, proc, + proc->waitLockMode, + &proc->waitLock->tag); early_deadlock = true; break; } diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h index ee3cb1dc203..85804d61ce9 100644 --- a/src/include/storage/lock.h +++ b/src/include/storage/lock.h @@ -401,6 +401,7 @@ extern void LockReleaseCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern void LockReassignCurrentOwner(LOCALLOCK **locallocks, int nlocks); extern bool LockHeldByMe(const LOCKTAG *locktag, LOCKMODE lockmode, bool orstronger); +extern bool HaveFastPathLock(Oid relid); #ifdef USE_ASSERT_CHECKING extern HTAB *GetLockMethodLocalHash(void); #endif @@ -440,7 +441,9 @@ pg_noreturn extern void DeadLockReport(void); extern void RememberSimpleDeadLock(PGPROC *proc1, LOCKMODE lockmode, LOCK *lock, - PGPROC *proc2); + PGPROC *proc2, + LOCKMODE lockmode2, + LOCKTAG *locktag2); extern void InitDeadLockChecking(void); extern int LockWaiterCount(const LOCKTAG *locktag); diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h index 3e1d1fad5f9..76c6bb44251 100644 --- a/src/include/storage/proc.h +++ b/src/include/storage/proc.h @@ -70,10 +70,12 @@ struct XidCache #define PROC_AFFECTS_ALL_HORIZONS 0x20 /* this proc's xmin must be * included in vacuum horizons * in all databases */ +#define PROC_IN_CONCURRENT_REPACK 0x40 /* REPACK (CONCURRENTLY) */ -/* flags reset at EOXact */ +/* flags reset at EOXact. A bit of a misnomer ... */ #define PROC_VACUUM_STATE_MASK \ - (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND) + (PROC_IN_VACUUM | PROC_IN_SAFE_IC | PROC_VACUUM_FOR_WRAPAROUND | \ + PROC_IN_CONCURRENT_REPACK) /* * Xmin-related flags. Make sure any flags that affect how the process' Xmin diff --git a/src/test/modules/injection_points/expected/repack.out b/src/test/modules/injection_points/expected/repack.out index b575e9052ee..960044776f4 100644 --- a/src/test/modules/injection_points/expected/repack.out +++ b/src/test/modules/injection_points/expected/repack.out @@ -1,4 +1,4 @@ -Parsed test spec with 2 sessions +Parsed test spec with 3 sessions starting permutation: wait_before_lock change_existing change_new change_subxact1 change_subxact2 check2 wakeup_before_lock check1 injection_points_attach @@ -111,3 +111,60 @@ injection_points_detach (1 row) + +starting permutation: check1_relnode_only wait_before_lock lock3 wakeup_before_lock check1_relnode_only +injection_points_attach +----------------------- + +(1 row) + +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 1 +(1 row) + +step wait_before_lock: + REPACK (CONCURRENTLY) repack_test USING INDEX repack_test_pkey; + <waiting ...> +step lock3: + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; + +i|j +-+- +1|1 +(1 row) + +ERROR: deadlock detected +step wakeup_before_lock: + SELECT injection_points_wakeup('repack-concurrently-before-lock'); + +injection_points_wakeup +----------------------- + +(1 row) + +step wait_before_lock: <... completed> +step check1_relnode_only: + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; + +count +----- + 2 +(1 row) + +injection_points_detach +----------------------- + +(1 row) + diff --git a/src/test/modules/injection_points/specs/repack.spec b/src/test/modules/injection_points/specs/repack.spec index d727a9b056b..5ffcb558718 100644 --- a/src/test/modules/injection_points/specs/repack.spec +++ b/src/test/modules/injection_points/specs/repack.spec @@ -55,6 +55,13 @@ step check1 FROM data_s1 d1 FULL JOIN data_s2 d2 USING (i, j) WHERE d1.i ISNULL OR d2.i ISNULL; } +step check1_relnode_only +{ + INSERT INTO relfilenodes(node) + SELECT relfilenode FROM pg_class WHERE relname='repack_test'; + + SELECT count(DISTINCT node) FROM relfilenodes; +} teardown { SELECT injection_points_detach('repack-concurrently-before-lock'); @@ -129,6 +136,20 @@ step wakeup_before_lock SELECT injection_points_wakeup('repack-concurrently-before-lock'); } +session s3 +# Try to acquire SharedUpdateExclusiveLock on a table while REPACK is already +# holding it and before it tries to upgrade it to AccessExclusiveLock. Since +# this session is already holding another lock no the table, REPACK cannot +# resolve the problem by getting ahead in the wait queue. The solution is that +# the LOCK TABLE triggers a deadlock report. The deadlock does not actually +# take place, but it would if this session started to wait. +step lock3 +{ + BEGIN; + SELECT * FROM repack_test ORDER BY i LIMIT 1; + LOCK TABLE repack_test IN SHARE UPDATE EXCLUSIVE MODE; +} + # Test if data changes introduced while one session is performing REPACK # CONCURRENTLY find their way into the table. permutation @@ -140,3 +161,11 @@ permutation check2 wakeup_before_lock check1 + +# See 'lock3' above. We check relnodes to make sure that REPACK finished. +permutation + check1_relnode_only + wait_before_lock + lock3 + wakeup_before_lock + check1_relnode_only -- 2.47.3 --=-=-=-- ^ permalink raw reply [nested|flat] 386+ messages in thread
end of thread, other threads:[~2026-04-14 09:59 UTC | newest] Thread overview: 386+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2025-08-31 09:10 [PATCH v1 4/4] Extra tests Nikolay Shaplov <dhyan@nataraj.su> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at> 2026-04-14 09:59 [PATCH] Teach REPACK to upgrade its lock safely. Antonin Houska <ah@cybertec.at>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox